agora inbox for [email protected]  
help / color / mirror / Atom feed
[PATCH v19 1/6] Split vacuumdb to create vacuuming.c/h
315+ messages / 2 participants
[nested] [flat]

* [PATCH v19 1/6] Split vacuumdb to create vacuuming.c/h
@ 2025-08-30 12:39 Álvaro Herrera <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Álvaro Herrera @ 2025-08-30 12:39 UTC (permalink / raw)

---
 src/bin/scripts/Makefile    |    4 +-
 src/bin/scripts/meson.build |   28 +-
 src/bin/scripts/vacuumdb.c  | 1048 +----------------------------------
 src/bin/scripts/vacuuming.c |  978 ++++++++++++++++++++++++++++++++
 src/bin/scripts/vacuuming.h |   95 ++++
 5 files changed, 1119 insertions(+), 1034 deletions(-)
 create mode 100644 src/bin/scripts/vacuuming.c
 create mode 100644 src/bin/scripts/vacuuming.h

diff --git a/src/bin/scripts/Makefile b/src/bin/scripts/Makefile
index f6b4d40810b..019ca06455d 100644
--- a/src/bin/scripts/Makefile
+++ b/src/bin/scripts/Makefile
@@ -28,7 +28,7 @@ createuser: createuser.o common.o $(WIN32RES) | submake-libpq submake-libpgport
 dropdb: dropdb.o common.o $(WIN32RES) | submake-libpq submake-libpgport submake-libpgfeutils
 dropuser: dropuser.o common.o $(WIN32RES) | submake-libpq submake-libpgport submake-libpgfeutils
 clusterdb: clusterdb.o common.o $(WIN32RES) | submake-libpq submake-libpgport submake-libpgfeutils
-vacuumdb: vacuumdb.o common.o $(WIN32RES) | submake-libpq submake-libpgport submake-libpgfeutils
+vacuumdb: vacuumdb.o vacuuming.o common.o $(WIN32RES) | submake-libpq submake-libpgport submake-libpgfeutils
 reindexdb: reindexdb.o common.o $(WIN32RES) | submake-libpq submake-libpgport submake-libpgfeutils
 pg_isready: pg_isready.o common.o $(WIN32RES) | submake-libpq submake-libpgport submake-libpgfeutils
 
@@ -50,7 +50,7 @@ uninstall:
 
 clean distclean:
 	rm -f $(addsuffix $(X), $(PROGRAMS)) $(addsuffix .o, $(PROGRAMS))
-	rm -f common.o $(WIN32RES)
+	rm -f common.o vacuuming.o $(WIN32RES)
 	rm -rf tmp_check
 
 export with_icu
diff --git a/src/bin/scripts/meson.build b/src/bin/scripts/meson.build
index 80df7c33257..a4fed59d1c9 100644
--- a/src/bin/scripts/meson.build
+++ b/src/bin/scripts/meson.build
@@ -12,7 +12,6 @@ binaries = [
   'createuser',
   'dropuser',
   'clusterdb',
-  'vacuumdb',
   'reindexdb',
   'pg_isready',
 ]
@@ -35,6 +34,33 @@ foreach binary : binaries
   bin_targets += binary
 endforeach
 
+vacuuming_common = static_library('libvacuuming_common',
+  files('common.c', 'vacuuming.c'),
+  dependencies: [frontend_code, libpq],
+  kwargs: internal_lib_args,
+)
+
+binaries = [
+  'vacuumdb',
+]
+foreach binary : binaries
+  binary_sources = files('@[email protected]'.format(binary))
+
+  if host_system == 'windows'
+    binary_sources += rc_bin_gen.process(win32ver_rc, extra_args: [
+      '--NAME', binary,
+      '--FILEDESC', '@0@ - PostgreSQL utility'.format(binary),])
+  endif
+
+  binary = executable(binary,
+    binary_sources,
+    link_with: [vacuuming_common],
+    dependencies: [frontend_code, libpq],
+    kwargs: default_bin_args,
+  )
+  bin_targets += binary
+endforeach
+
 tests += {
   'name': 'scripts',
   'sd': meson.current_source_dir(),
diff --git a/src/bin/scripts/vacuumdb.c b/src/bin/scripts/vacuumdb.c
index fd236087e90..b1be61ddf25 100644
--- a/src/bin/scripts/vacuumdb.c
+++ b/src/bin/scripts/vacuumdb.c
@@ -14,92 +14,13 @@
 
 #include <limits.h>
 
-#include "catalog/pg_attribute_d.h"
-#include "catalog/pg_class_d.h"
 #include "common.h"
-#include "common/connect.h"
 #include "common/logging.h"
-#include "fe_utils/cancel.h"
 #include "fe_utils/option_utils.h"
-#include "fe_utils/parallel_slot.h"
-#include "fe_utils/query_utils.h"
-#include "fe_utils/simple_list.h"
-#include "fe_utils/string_utils.h"
-
-
-/* vacuum options controlled by user flags */
-typedef struct vacuumingOptions
-{
-	bool		analyze_only;
-	bool		verbose;
-	bool		and_analyze;
-	bool		full;
-	bool		freeze;
-	bool		disable_page_skipping;
-	bool		skip_locked;
-	int			min_xid_age;
-	int			min_mxid_age;
-	int			parallel_workers;	/* >= 0 indicates user specified the
-									 * parallel degree, otherwise -1 */
-	bool		no_index_cleanup;
-	bool		force_index_cleanup;
-	bool		do_truncate;
-	bool		process_main;
-	bool		process_toast;
-	bool		skip_database_stats;
-	char	   *buffer_usage_limit;
-	bool		missing_stats_only;
-} vacuumingOptions;
-
-/* object filter options */
-typedef enum
-{
-	OBJFILTER_NONE = 0,			/* no filter used */
-	OBJFILTER_ALL_DBS = (1 << 0),	/* -a | --all */
-	OBJFILTER_DATABASE = (1 << 1),	/* -d | --dbname */
-	OBJFILTER_TABLE = (1 << 2), /* -t | --table */
-	OBJFILTER_SCHEMA = (1 << 3),	/* -n | --schema */
-	OBJFILTER_SCHEMA_EXCLUDE = (1 << 4),	/* -N | --exclude-schema */
-} VacObjFilter;
-
-static VacObjFilter objfilter = OBJFILTER_NONE;
-
-static SimpleStringList *retrieve_objects(PGconn *conn,
-										  vacuumingOptions *vacopts,
-										  SimpleStringList *objects,
-										  bool echo);
-
-static void vacuum_one_database(ConnParams *cparams,
-								vacuumingOptions *vacopts,
-								int stage,
-								SimpleStringList *objects,
-								SimpleStringList **found_objs,
-								int concurrentCons,
-								const char *progname, bool echo, bool quiet);
-
-static void vacuum_all_databases(ConnParams *cparams,
-								 vacuumingOptions *vacopts,
-								 bool analyze_in_stages,
-								 SimpleStringList *objects,
-								 int concurrentCons,
-								 const char *progname, bool echo, bool quiet);
-
-static void prepare_vacuum_command(PQExpBuffer sql, int serverVersion,
-								   vacuumingOptions *vacopts, const char *table);
-
-static void run_vacuum_command(PGconn *conn, const char *sql, bool echo,
-							   const char *table);
+#include "vacuuming.h"
 
 static void help(const char *progname);
-
-void		check_objfilter(void);
-
-static char *escape_quotes(const char *src);
-
-/* For analyze-in-stages mode */
-#define ANALYZE_NO_STAGE	-1
-#define ANALYZE_NUM_STAGES	3
-
+static void check_objfilter(void);
 
 int
 main(int argc, char *argv[])
@@ -145,10 +66,6 @@ main(int argc, char *argv[])
 	int			c;
 	const char *dbname = NULL;
 	const char *maintenance_db = NULL;
-	char	   *host = NULL;
-	char	   *port = NULL;
-	char	   *username = NULL;
-	enum trivalue prompt_password = TRI_DEFAULT;
 	ConnParams	cparams;
 	bool		echo = false;
 	bool		quiet = false;
@@ -168,13 +85,18 @@ main(int argc, char *argv[])
 	vacopts.process_main = true;
 	vacopts.process_toast = true;
 
+	/* the same for connection parameters */
+	memset(&cparams, 0, sizeof(cparams));
+	cparams.prompt_password = TRI_DEFAULT;
+
 	pg_logging_init(argv[0]);
 	progname = get_progname(argv[0]);
 	set_pglocale_pgservice(argv[0], PG_TEXTDOMAIN("pgscripts"));
 
-	handle_help_version_opts(argc, argv, "vacuumdb", help);
+	handle_help_version_opts(argc, argv, progname, help);
 
-	while ((c = getopt_long(argc, argv, "ad:efFh:j:n:N:p:P:qt:U:vwWzZ", long_options, &optindex)) != -1)
+	while ((c = getopt_long(argc, argv, "ad:efFh:j:n:N:p:P:qt:U:vwWzZ",
+							long_options, &optindex)) != -1)
 	{
 		switch (c)
 		{
@@ -195,7 +117,7 @@ main(int argc, char *argv[])
 				vacopts.freeze = true;
 				break;
 			case 'h':
-				host = pg_strdup(optarg);
+				cparams.pghost = pg_strdup(optarg);
 				break;
 			case 'j':
 				if (!option_parse_int(optarg, "-j/--jobs", 1, INT_MAX,
@@ -211,7 +133,7 @@ main(int argc, char *argv[])
 				simple_string_list_append(&objects, optarg);
 				break;
 			case 'p':
-				port = pg_strdup(optarg);
+				cparams.pgport = pg_strdup(optarg);
 				break;
 			case 'P':
 				if (!option_parse_int(optarg, "-P/--parallel", 0, INT_MAX,
@@ -227,16 +149,16 @@ main(int argc, char *argv[])
 				tbl_count++;
 				break;
 			case 'U':
-				username = pg_strdup(optarg);
+				cparams.pguser = pg_strdup(optarg);
 				break;
 			case 'v':
 				vacopts.verbose = true;
 				break;
 			case 'w':
-				prompt_password = TRI_NO;
+				cparams.prompt_password = TRI_NO;
 				break;
 			case 'W':
-				prompt_password = TRI_YES;
+				cparams.prompt_password = TRI_YES;
 				break;
 			case 'z':
 				vacopts.and_analyze = true;
@@ -380,66 +302,9 @@ main(int argc, char *argv[])
 		pg_fatal("cannot use the \"%s\" option without \"%s\" or \"%s\"",
 				 "missing-stats-only", "analyze-only", "analyze-in-stages");
 
-	/* fill cparams except for dbname, which is set below */
-	cparams.pghost = host;
-	cparams.pgport = port;
-	cparams.pguser = username;
-	cparams.prompt_password = prompt_password;
-	cparams.override_dbname = NULL;
-
-	setup_cancel_handler(NULL);
-
-	/* Avoid opening extra connections. */
-	if (tbl_count && (concurrentCons > tbl_count))
-		concurrentCons = tbl_count;
-
-	if (objfilter & OBJFILTER_ALL_DBS)
-	{
-		cparams.dbname = maintenance_db;
-
-		vacuum_all_databases(&cparams, &vacopts,
-							 analyze_in_stages,
-							 &objects,
-							 concurrentCons,
-							 progname, echo, quiet);
-	}
-	else
-	{
-		if (dbname == NULL)
-		{
-			if (getenv("PGDATABASE"))
-				dbname = getenv("PGDATABASE");
-			else if (getenv("PGUSER"))
-				dbname = getenv("PGUSER");
-			else
-				dbname = get_user_name_or_exit(progname);
-		}
-
-		cparams.dbname = dbname;
-
-		if (analyze_in_stages)
-		{
-			int			stage;
-			SimpleStringList *found_objs = NULL;
-
-			for (stage = 0; stage < ANALYZE_NUM_STAGES; stage++)
-			{
-				vacuum_one_database(&cparams, &vacopts,
-									stage,
-									&objects,
-									vacopts.missing_stats_only ? &found_objs : NULL,
-									concurrentCons,
-									progname, echo, quiet);
-			}
-		}
-		else
-			vacuum_one_database(&cparams, &vacopts,
-								ANALYZE_NO_STAGE,
-								&objects, NULL,
-								concurrentCons,
-								progname, echo, quiet);
-	}
-
+	vacuuming_main(&cparams, dbname, maintenance_db, &vacopts, &objects,
+				   analyze_in_stages, tbl_count, concurrentCons,
+				   progname, echo, quiet);
 	exit(0);
 }
 
@@ -466,885 +331,6 @@ check_objfilter(void)
 		pg_fatal("cannot vacuum all tables in schema(s) and exclude schema(s) at the same time");
 }
 
-/*
- * Returns a newly malloc'd version of 'src' with escaped single quotes and
- * backslashes.
- */
-static char *
-escape_quotes(const char *src)
-{
-	char	   *result = escape_single_quotes_ascii(src);
-
-	if (!result)
-		pg_fatal("out of memory");
-	return result;
-}
-
-/*
- * vacuum_one_database
- *
- * Process tables in the given database.
- *
- * There are two ways to specify the list of objects to process:
- *
- * 1) The "found_objs" parameter is a double pointer to a fully qualified list
- *    of objects to process, as returned by a previous call to
- *    vacuum_one_database().
- *
- *     a) If both "found_objs" (the double pointer) and "*found_objs" (the
- *        once-dereferenced double pointer) are not NULL, this list takes
- *        priority, and anything specified in "objects" is ignored.
- *
- *     b) If "found_objs" (the double pointer) is not NULL but "*found_objs"
- *        (the once-dereferenced double pointer) _is_ NULL, the "objects"
- *        parameter takes priority, and the results of the catalog query
- *        described in (2) are stored in "found_objs".
- *
- *     c) If "found_objs" (the double pointer) is NULL, the "objects"
- *        parameter again takes priority, and the results of the catalog query
- *        are not saved.
- *
- * 2) The "objects" parameter is a user-specified list of objects to process.
- *    When (1b) or (1c) applies, this function performs a catalog query to
- *    retrieve a fully qualified list of objects to process, as described
- *    below.
- *
- *     a) If "objects" is not NULL, the catalog query gathers only the objects
- *        listed in "objects".
- *
- *     b) If "objects" is NULL, all tables in the database are gathered.
- *
- * Note that this function is only concerned with running exactly one stage
- * when in analyze-in-stages mode; caller must iterate on us if necessary.
- *
- * If concurrentCons is > 1, multiple connections are used to vacuum tables
- * in parallel.
- */
-static void
-vacuum_one_database(ConnParams *cparams,
-					vacuumingOptions *vacopts,
-					int stage,
-					SimpleStringList *objects,
-					SimpleStringList **found_objs,
-					int concurrentCons,
-					const char *progname, bool echo, bool quiet)
-{
-	PQExpBufferData sql;
-	PGconn	   *conn;
-	SimpleStringListCell *cell;
-	ParallelSlotArray *sa;
-	int			ntups = 0;
-	bool		failed = false;
-	const char *initcmd;
-	SimpleStringList *ret = NULL;
-	const char *stage_commands[] = {
-		"SET default_statistics_target=1; SET vacuum_cost_delay=0;",
-		"SET default_statistics_target=10; RESET vacuum_cost_delay;",
-		"RESET default_statistics_target;"
-	};
-	const char *stage_messages[] = {
-		gettext_noop("Generating minimal optimizer statistics (1 target)"),
-		gettext_noop("Generating medium optimizer statistics (10 targets)"),
-		gettext_noop("Generating default (full) optimizer statistics")
-	};
-
-	Assert(stage == ANALYZE_NO_STAGE ||
-		   (stage >= 0 && stage < ANALYZE_NUM_STAGES));
-
-	conn = connectDatabase(cparams, progname, echo, false, true);
-
-	if (vacopts->disable_page_skipping && PQserverVersion(conn) < 90600)
-	{
-		PQfinish(conn);
-		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
-				 "disable-page-skipping", "9.6");
-	}
-
-	if (vacopts->no_index_cleanup && PQserverVersion(conn) < 120000)
-	{
-		PQfinish(conn);
-		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
-				 "no-index-cleanup", "12");
-	}
-
-	if (vacopts->force_index_cleanup && PQserverVersion(conn) < 120000)
-	{
-		PQfinish(conn);
-		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
-				 "force-index-cleanup", "12");
-	}
-
-	if (!vacopts->do_truncate && PQserverVersion(conn) < 120000)
-	{
-		PQfinish(conn);
-		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
-				 "no-truncate", "12");
-	}
-
-	if (!vacopts->process_main && PQserverVersion(conn) < 160000)
-	{
-		PQfinish(conn);
-		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
-				 "no-process-main", "16");
-	}
-
-	if (!vacopts->process_toast && PQserverVersion(conn) < 140000)
-	{
-		PQfinish(conn);
-		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
-				 "no-process-toast", "14");
-	}
-
-	if (vacopts->skip_locked && PQserverVersion(conn) < 120000)
-	{
-		PQfinish(conn);
-		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
-				 "skip-locked", "12");
-	}
-
-	if (vacopts->min_xid_age != 0 && PQserverVersion(conn) < 90600)
-	{
-		PQfinish(conn);
-		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
-				 "--min-xid-age", "9.6");
-	}
-
-	if (vacopts->min_mxid_age != 0 && PQserverVersion(conn) < 90600)
-	{
-		PQfinish(conn);
-		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
-				 "--min-mxid-age", "9.6");
-	}
-
-	if (vacopts->parallel_workers >= 0 && PQserverVersion(conn) < 130000)
-	{
-		PQfinish(conn);
-		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
-				 "--parallel", "13");
-	}
-
-	if (vacopts->buffer_usage_limit && PQserverVersion(conn) < 160000)
-	{
-		PQfinish(conn);
-		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
-				 "--buffer-usage-limit", "16");
-	}
-
-	if (vacopts->missing_stats_only && PQserverVersion(conn) < 150000)
-	{
-		PQfinish(conn);
-		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
-				 "--missing-stats-only", "15");
-	}
-
-	/* skip_database_stats is used automatically if server supports it */
-	vacopts->skip_database_stats = (PQserverVersion(conn) >= 160000);
-
-	if (!quiet)
-	{
-		if (stage != ANALYZE_NO_STAGE)
-			printf(_("%s: processing database \"%s\": %s\n"),
-				   progname, PQdb(conn), _(stage_messages[stage]));
-		else
-			printf(_("%s: vacuuming database \"%s\"\n"),
-				   progname, PQdb(conn));
-		fflush(stdout);
-	}
-
-	/*
-	 * If the caller provided the results of a previous catalog query, just
-	 * use that.  Otherwise, run the catalog query ourselves and set the
-	 * return variable if provided.
-	 */
-	if (found_objs && *found_objs)
-		ret = *found_objs;
-	else
-	{
-		ret = retrieve_objects(conn, vacopts, objects, echo);
-		if (found_objs)
-			*found_objs = ret;
-	}
-
-	/*
-	 * Count the number of objects in the catalog query result.  If there are
-	 * none, we are done.
-	 */
-	for (cell = ret ? ret->head : NULL; cell; cell = cell->next)
-		ntups++;
-
-	if (ntups == 0)
-	{
-		PQfinish(conn);
-		return;
-	}
-
-	/*
-	 * Ensure concurrentCons is sane.  If there are more connections than
-	 * vacuumable relations, we don't need to use them all.
-	 */
-	if (concurrentCons > ntups)
-		concurrentCons = ntups;
-	if (concurrentCons <= 0)
-		concurrentCons = 1;
-
-	/*
-	 * All slots need to be prepared to run the appropriate analyze stage, if
-	 * caller requested that mode.  We have to prepare the initial connection
-	 * ourselves before setting up the slots.
-	 */
-	if (stage == ANALYZE_NO_STAGE)
-		initcmd = NULL;
-	else
-	{
-		initcmd = stage_commands[stage];
-		executeCommand(conn, initcmd, echo);
-	}
-
-	/*
-	 * Setup the database connections. We reuse the connection we already have
-	 * for the first slot.  If not in parallel mode, the first slot in the
-	 * array contains the connection.
-	 */
-	sa = ParallelSlotsSetup(concurrentCons, cparams, progname, echo, initcmd);
-	ParallelSlotsAdoptConn(sa, conn);
-
-	initPQExpBuffer(&sql);
-
-	cell = ret->head;
-	do
-	{
-		const char *tabname = cell->val;
-		ParallelSlot *free_slot;
-
-		if (CancelRequested)
-		{
-			failed = true;
-			goto finish;
-		}
-
-		free_slot = ParallelSlotsGetIdle(sa, NULL);
-		if (!free_slot)
-		{
-			failed = true;
-			goto finish;
-		}
-
-		prepare_vacuum_command(&sql, PQserverVersion(free_slot->connection),
-							   vacopts, tabname);
-
-		/*
-		 * Execute the vacuum.  All errors are handled in processQueryResult
-		 * through ParallelSlotsGetIdle.
-		 */
-		ParallelSlotSetHandler(free_slot, TableCommandResultHandler, NULL);
-		run_vacuum_command(free_slot->connection, sql.data,
-						   echo, tabname);
-
-		cell = cell->next;
-	} while (cell != NULL);
-
-	if (!ParallelSlotsWaitCompletion(sa))
-	{
-		failed = true;
-		goto finish;
-	}
-
-	/* If we used SKIP_DATABASE_STATS, mop up with ONLY_DATABASE_STATS */
-	if (vacopts->skip_database_stats && stage == ANALYZE_NO_STAGE)
-	{
-		const char *cmd = "VACUUM (ONLY_DATABASE_STATS);";
-		ParallelSlot *free_slot = ParallelSlotsGetIdle(sa, NULL);
-
-		if (!free_slot)
-		{
-			failed = true;
-			goto finish;
-		}
-
-		ParallelSlotSetHandler(free_slot, TableCommandResultHandler, NULL);
-		run_vacuum_command(free_slot->connection, cmd, echo, NULL);
-
-		if (!ParallelSlotsWaitCompletion(sa))
-			failed = true;
-	}
-
-finish:
-	ParallelSlotsTerminate(sa);
-	pg_free(sa);
-
-	termPQExpBuffer(&sql);
-
-	if (failed)
-		exit(1);
-}
-
-/*
- * Prepare the list of tables to process by querying the catalogs.
- *
- * Since we execute the constructed query with the default search_path (which
- * could be unsafe), everything in this query MUST be fully qualified.
- *
- * First, build a WITH clause for the catalog query if any tables were
- * specified, with a set of values made of relation names and their optional
- * set of columns.  This is used to match any provided column lists with the
- * generated qualified identifiers and to filter for the tables provided via
- * --table.  If a listed table does not exist, the catalog query will fail.
- */
-static SimpleStringList *
-retrieve_objects(PGconn *conn, vacuumingOptions *vacopts,
-				 SimpleStringList *objects, bool echo)
-{
-	PQExpBufferData buf;
-	PQExpBufferData catalog_query;
-	PGresult   *res;
-	SimpleStringListCell *cell;
-	SimpleStringList *found_objs = palloc0(sizeof(SimpleStringList));
-	bool		objects_listed = false;
-
-	initPQExpBuffer(&catalog_query);
-	for (cell = objects ? objects->head : NULL; cell; cell = cell->next)
-	{
-		char	   *just_table = NULL;
-		const char *just_columns = NULL;
-
-		if (!objects_listed)
-		{
-			appendPQExpBufferStr(&catalog_query,
-								 "WITH listed_objects (object_oid, column_list) "
-								 "AS (\n  VALUES (");
-			objects_listed = true;
-		}
-		else
-			appendPQExpBufferStr(&catalog_query, ",\n  (");
-
-		if (objfilter & (OBJFILTER_SCHEMA | OBJFILTER_SCHEMA_EXCLUDE))
-		{
-			appendStringLiteralConn(&catalog_query, cell->val, conn);
-			appendPQExpBufferStr(&catalog_query, "::pg_catalog.regnamespace, ");
-		}
-
-		if (objfilter & OBJFILTER_TABLE)
-		{
-			/*
-			 * Split relation and column names given by the user, this is used
-			 * to feed the CTE with values on which are performed pre-run
-			 * validity checks as well.  For now these happen only on the
-			 * relation name.
-			 */
-			splitTableColumnsSpec(cell->val, PQclientEncoding(conn),
-								  &just_table, &just_columns);
-
-			appendStringLiteralConn(&catalog_query, just_table, conn);
-			appendPQExpBufferStr(&catalog_query, "::pg_catalog.regclass, ");
-		}
-
-		if (just_columns && just_columns[0] != '\0')
-			appendStringLiteralConn(&catalog_query, just_columns, conn);
-		else
-			appendPQExpBufferStr(&catalog_query, "NULL");
-
-		appendPQExpBufferStr(&catalog_query, "::pg_catalog.text)");
-
-		pg_free(just_table);
-	}
-
-	/* Finish formatting the CTE */
-	if (objects_listed)
-		appendPQExpBufferStr(&catalog_query, "\n)\n");
-
-	appendPQExpBufferStr(&catalog_query, "SELECT c.relname, ns.nspname");
-
-	if (objects_listed)
-		appendPQExpBufferStr(&catalog_query, ", listed_objects.column_list");
-
-	appendPQExpBufferStr(&catalog_query,
-						 " FROM pg_catalog.pg_class c\n"
-						 " JOIN pg_catalog.pg_namespace ns"
-						 " ON c.relnamespace OPERATOR(pg_catalog.=) ns.oid\n"
-						 " CROSS JOIN LATERAL (SELECT c.relkind IN ("
-						 CppAsString2(RELKIND_PARTITIONED_TABLE) ", "
-						 CppAsString2(RELKIND_PARTITIONED_INDEX) ")) as p (inherited)\n"
-						 " LEFT JOIN pg_catalog.pg_class t"
-						 " ON c.reltoastrelid OPERATOR(pg_catalog.=) t.oid\n");
-
-	/*
-	 * Used to match the tables or schemas listed by the user, completing the
-	 * JOIN clause.
-	 */
-	if (objects_listed)
-	{
-		appendPQExpBufferStr(&catalog_query, " LEFT JOIN listed_objects"
-							 " ON listed_objects.object_oid"
-							 " OPERATOR(pg_catalog.=) ");
-
-		if (objfilter & OBJFILTER_TABLE)
-			appendPQExpBufferStr(&catalog_query, "c.oid\n");
-		else
-			appendPQExpBufferStr(&catalog_query, "ns.oid\n");
-	}
-
-	/*
-	 * Exclude temporary tables, beginning the WHERE clause.
-	 */
-	appendPQExpBufferStr(&catalog_query,
-						 " WHERE c.relpersistence OPERATOR(pg_catalog.!=) "
-						 CppAsString2(RELPERSISTENCE_TEMP) "\n");
-
-	/*
-	 * Used to match the tables or schemas listed by the user, for the WHERE
-	 * clause.
-	 */
-	if (objects_listed)
-	{
-		if (objfilter & OBJFILTER_SCHEMA_EXCLUDE)
-			appendPQExpBufferStr(&catalog_query,
-								 " AND listed_objects.object_oid IS NULL\n");
-		else
-			appendPQExpBufferStr(&catalog_query,
-								 " AND listed_objects.object_oid IS NOT NULL\n");
-	}
-
-	/*
-	 * If no tables were listed, filter for the relevant relation types.  If
-	 * tables were given via --table, don't bother filtering by relation type.
-	 * Instead, let the server decide whether a given relation can be
-	 * processed in which case the user will know about it.
-	 */
-	if ((objfilter & OBJFILTER_TABLE) == 0)
-	{
-		/*
-		 * vacuumdb should generally follow the behavior of the underlying
-		 * VACUUM and ANALYZE commands. If analyze_only is true, process
-		 * regular tables, materialized views, and partitioned tables, just
-		 * like ANALYZE (with no specific target tables) does. Otherwise,
-		 * process only regular tables and materialized views, since VACUUM
-		 * skips partitioned tables when no target tables are specified.
-		 */
-		if (vacopts->analyze_only)
-			appendPQExpBufferStr(&catalog_query,
-								 " AND c.relkind OPERATOR(pg_catalog.=) ANY (array["
-								 CppAsString2(RELKIND_RELATION) ", "
-								 CppAsString2(RELKIND_MATVIEW) ", "
-								 CppAsString2(RELKIND_PARTITIONED_TABLE) "])\n");
-		else
-			appendPQExpBufferStr(&catalog_query,
-								 " AND c.relkind OPERATOR(pg_catalog.=) ANY (array["
-								 CppAsString2(RELKIND_RELATION) ", "
-								 CppAsString2(RELKIND_MATVIEW) "])\n");
-
-	}
-
-	/*
-	 * For --min-xid-age and --min-mxid-age, the age of the relation is the
-	 * greatest of the ages of the main relation and its associated TOAST
-	 * table.  The commands generated by vacuumdb will also process the TOAST
-	 * table for the relation if necessary, so it does not need to be
-	 * considered separately.
-	 */
-	if (vacopts->min_xid_age != 0)
-	{
-		appendPQExpBuffer(&catalog_query,
-						  " AND GREATEST(pg_catalog.age(c.relfrozenxid),"
-						  " pg_catalog.age(t.relfrozenxid)) "
-						  " OPERATOR(pg_catalog.>=) '%d'::pg_catalog.int4\n"
-						  " AND c.relfrozenxid OPERATOR(pg_catalog.!=)"
-						  " '0'::pg_catalog.xid\n",
-						  vacopts->min_xid_age);
-	}
-
-	if (vacopts->min_mxid_age != 0)
-	{
-		appendPQExpBuffer(&catalog_query,
-						  " AND GREATEST(pg_catalog.mxid_age(c.relminmxid),"
-						  " pg_catalog.mxid_age(t.relminmxid)) OPERATOR(pg_catalog.>=)"
-						  " '%d'::pg_catalog.int4\n"
-						  " AND c.relminmxid OPERATOR(pg_catalog.!=)"
-						  " '0'::pg_catalog.xid\n",
-						  vacopts->min_mxid_age);
-	}
-
-	if (vacopts->missing_stats_only)
-	{
-		appendPQExpBufferStr(&catalog_query, " AND (\n");
-
-		/* regular stats */
-		appendPQExpBufferStr(&catalog_query,
-							 " EXISTS (SELECT NULL FROM pg_catalog.pg_attribute a\n"
-							 " WHERE a.attrelid OPERATOR(pg_catalog.=) c.oid\n"
-							 " AND a.attnum OPERATOR(pg_catalog.>) 0::pg_catalog.int2\n"
-							 " AND NOT a.attisdropped\n"
-							 " AND a.attstattarget IS DISTINCT FROM 0::pg_catalog.int2\n"
-							 " AND a.attgenerated OPERATOR(pg_catalog.<>) "
-							 CppAsString2(ATTRIBUTE_GENERATED_VIRTUAL) "\n"
-							 " AND NOT EXISTS (SELECT NULL FROM pg_catalog.pg_statistic s\n"
-							 " WHERE s.starelid OPERATOR(pg_catalog.=) a.attrelid\n"
-							 " AND s.staattnum OPERATOR(pg_catalog.=) a.attnum\n"
-							 " AND s.stainherit OPERATOR(pg_catalog.=) p.inherited))\n");
-
-		/* extended stats */
-		appendPQExpBufferStr(&catalog_query,
-							 " OR EXISTS (SELECT NULL FROM pg_catalog.pg_statistic_ext e\n"
-							 " WHERE e.stxrelid OPERATOR(pg_catalog.=) c.oid\n"
-							 " AND e.stxstattarget IS DISTINCT FROM 0::pg_catalog.int2\n"
-							 " AND NOT EXISTS (SELECT NULL FROM pg_catalog.pg_statistic_ext_data d\n"
-							 " WHERE d.stxoid OPERATOR(pg_catalog.=) e.oid\n"
-							 " AND d.stxdinherit OPERATOR(pg_catalog.=) p.inherited))\n");
-
-		/* expression indexes */
-		appendPQExpBufferStr(&catalog_query,
-							 " OR EXISTS (SELECT NULL FROM pg_catalog.pg_attribute a\n"
-							 " JOIN pg_catalog.pg_index i"
-							 " ON i.indexrelid OPERATOR(pg_catalog.=) a.attrelid\n"
-							 " WHERE i.indrelid OPERATOR(pg_catalog.=) c.oid\n"
-							 " AND i.indkey[a.attnum OPERATOR(pg_catalog.-) 1::pg_catalog.int2]"
-							 " OPERATOR(pg_catalog.=) 0::pg_catalog.int2\n"
-							 " AND a.attnum OPERATOR(pg_catalog.>) 0::pg_catalog.int2\n"
-							 " AND NOT a.attisdropped\n"
-							 " AND a.attstattarget IS DISTINCT FROM 0::pg_catalog.int2\n"
-							 " AND NOT EXISTS (SELECT NULL FROM pg_catalog.pg_statistic s\n"
-							 " WHERE s.starelid OPERATOR(pg_catalog.=) a.attrelid\n"
-							 " AND s.staattnum OPERATOR(pg_catalog.=) a.attnum\n"
-							 " AND s.stainherit OPERATOR(pg_catalog.=) p.inherited))\n");
-
-		/* inheritance and regular stats */
-		appendPQExpBufferStr(&catalog_query,
-							 " OR EXISTS (SELECT NULL FROM pg_catalog.pg_attribute a\n"
-							 " WHERE a.attrelid OPERATOR(pg_catalog.=) c.oid\n"
-							 " AND a.attnum OPERATOR(pg_catalog.>) 0::pg_catalog.int2\n"
-							 " AND NOT a.attisdropped\n"
-							 " AND a.attstattarget IS DISTINCT FROM 0::pg_catalog.int2\n"
-							 " AND a.attgenerated OPERATOR(pg_catalog.<>) "
-							 CppAsString2(ATTRIBUTE_GENERATED_VIRTUAL) "\n"
-							 " AND c.relhassubclass\n"
-							 " AND NOT p.inherited\n"
-							 " AND EXISTS (SELECT NULL FROM pg_catalog.pg_inherits h\n"
-							 " WHERE h.inhparent OPERATOR(pg_catalog.=) c.oid)\n"
-							 " AND NOT EXISTS (SELECT NULL FROM pg_catalog.pg_statistic s\n"
-							 " WHERE s.starelid OPERATOR(pg_catalog.=) a.attrelid\n"
-							 " AND s.staattnum OPERATOR(pg_catalog.=) a.attnum\n"
-							 " AND s.stainherit))\n");
-
-		/* inheritance and extended stats */
-		appendPQExpBufferStr(&catalog_query,
-							 " OR EXISTS (SELECT NULL FROM pg_catalog.pg_statistic_ext e\n"
-							 " WHERE e.stxrelid OPERATOR(pg_catalog.=) c.oid\n"
-							 " AND e.stxstattarget IS DISTINCT FROM 0::pg_catalog.int2\n"
-							 " AND c.relhassubclass\n"
-							 " AND NOT p.inherited\n"
-							 " AND EXISTS (SELECT NULL FROM pg_catalog.pg_inherits h\n"
-							 " WHERE h.inhparent OPERATOR(pg_catalog.=) c.oid)\n"
-							 " AND NOT EXISTS (SELECT NULL FROM pg_catalog.pg_statistic_ext_data d\n"
-							 " WHERE d.stxoid OPERATOR(pg_catalog.=) e.oid\n"
-							 " AND d.stxdinherit))\n");
-
-		appendPQExpBufferStr(&catalog_query, " )\n");
-	}
-
-	/*
-	 * Execute the catalog query.  We use the default search_path for this
-	 * query for consistency with table lookups done elsewhere by the user.
-	 */
-	appendPQExpBufferStr(&catalog_query, " ORDER BY c.relpages DESC;");
-	executeCommand(conn, "RESET search_path;", echo);
-	res = executeQuery(conn, catalog_query.data, echo);
-	termPQExpBuffer(&catalog_query);
-	PQclear(executeQuery(conn, ALWAYS_SECURE_SEARCH_PATH_SQL, echo));
-
-	/*
-	 * Build qualified identifiers for each table, including the column list
-	 * if given.
-	 */
-	initPQExpBuffer(&buf);
-	for (int i = 0; i < PQntuples(res); i++)
-	{
-		appendPQExpBufferStr(&buf,
-							 fmtQualifiedIdEnc(PQgetvalue(res, i, 1),
-											   PQgetvalue(res, i, 0),
-											   PQclientEncoding(conn)));
-
-		if (objects_listed && !PQgetisnull(res, i, 2))
-			appendPQExpBufferStr(&buf, PQgetvalue(res, i, 2));
-
-		simple_string_list_append(found_objs, buf.data);
-		resetPQExpBuffer(&buf);
-	}
-	termPQExpBuffer(&buf);
-	PQclear(res);
-
-	return found_objs;
-}
-
-/*
- * Vacuum/analyze all connectable databases.
- *
- * In analyze-in-stages mode, we process all databases in one stage before
- * moving on to the next stage.  That ensure minimal stats are available
- * quickly everywhere before generating more detailed ones.
- */
-static void
-vacuum_all_databases(ConnParams *cparams,
-					 vacuumingOptions *vacopts,
-					 bool analyze_in_stages,
-					 SimpleStringList *objects,
-					 int concurrentCons,
-					 const char *progname, bool echo, bool quiet)
-{
-	PGconn	   *conn;
-	PGresult   *result;
-	int			stage;
-	int			i;
-
-	conn = connectMaintenanceDatabase(cparams, progname, echo);
-	result = executeQuery(conn,
-						  "SELECT datname FROM pg_database WHERE datallowconn AND datconnlimit <> -2 ORDER BY 1;",
-						  echo);
-	PQfinish(conn);
-
-	if (analyze_in_stages)
-	{
-		SimpleStringList **found_objs = NULL;
-
-		if (vacopts->missing_stats_only)
-			found_objs = palloc0(PQntuples(result) * sizeof(SimpleStringList *));
-
-		/*
-		 * When analyzing all databases in stages, we analyze them all in the
-		 * fastest stage first, so that initial statistics become available
-		 * for all of them as soon as possible.
-		 *
-		 * This means we establish several times as many connections, but
-		 * that's a secondary consideration.
-		 */
-		for (stage = 0; stage < ANALYZE_NUM_STAGES; stage++)
-		{
-			for (i = 0; i < PQntuples(result); i++)
-			{
-				cparams->override_dbname = PQgetvalue(result, i, 0);
-
-				vacuum_one_database(cparams, vacopts,
-									stage,
-									objects,
-									vacopts->missing_stats_only ? &found_objs[i] : NULL,
-									concurrentCons,
-									progname, echo, quiet);
-			}
-		}
-	}
-	else
-	{
-		for (i = 0; i < PQntuples(result); i++)
-		{
-			cparams->override_dbname = PQgetvalue(result, i, 0);
-
-			vacuum_one_database(cparams, vacopts,
-								ANALYZE_NO_STAGE,
-								objects, NULL,
-								concurrentCons,
-								progname, echo, quiet);
-		}
-	}
-
-	PQclear(result);
-}
-
-/*
- * Construct a vacuum/analyze command to run based on the given options, in the
- * given string buffer, which may contain previous garbage.
- *
- * The table name used must be already properly quoted.  The command generated
- * depends on the server version involved and it is semicolon-terminated.
- */
-static void
-prepare_vacuum_command(PQExpBuffer sql, int serverVersion,
-					   vacuumingOptions *vacopts, const char *table)
-{
-	const char *paren = " (";
-	const char *comma = ", ";
-	const char *sep = paren;
-
-	resetPQExpBuffer(sql);
-
-	if (vacopts->analyze_only)
-	{
-		appendPQExpBufferStr(sql, "ANALYZE");
-
-		/* parenthesized grammar of ANALYZE is supported since v11 */
-		if (serverVersion >= 110000)
-		{
-			if (vacopts->skip_locked)
-			{
-				/* SKIP_LOCKED is supported since v12 */
-				Assert(serverVersion >= 120000);
-				appendPQExpBuffer(sql, "%sSKIP_LOCKED", sep);
-				sep = comma;
-			}
-			if (vacopts->verbose)
-			{
-				appendPQExpBuffer(sql, "%sVERBOSE", sep);
-				sep = comma;
-			}
-			if (vacopts->buffer_usage_limit)
-			{
-				Assert(serverVersion >= 160000);
-				appendPQExpBuffer(sql, "%sBUFFER_USAGE_LIMIT '%s'", sep,
-								  vacopts->buffer_usage_limit);
-				sep = comma;
-			}
-			if (sep != paren)
-				appendPQExpBufferChar(sql, ')');
-		}
-		else
-		{
-			if (vacopts->verbose)
-				appendPQExpBufferStr(sql, " VERBOSE");
-		}
-	}
-	else
-	{
-		appendPQExpBufferStr(sql, "VACUUM");
-
-		/* parenthesized grammar of VACUUM is supported since v9.0 */
-		if (serverVersion >= 90000)
-		{
-			if (vacopts->disable_page_skipping)
-			{
-				/* DISABLE_PAGE_SKIPPING is supported since v9.6 */
-				Assert(serverVersion >= 90600);
-				appendPQExpBuffer(sql, "%sDISABLE_PAGE_SKIPPING", sep);
-				sep = comma;
-			}
-			if (vacopts->no_index_cleanup)
-			{
-				/* "INDEX_CLEANUP FALSE" has been supported since v12 */
-				Assert(serverVersion >= 120000);
-				Assert(!vacopts->force_index_cleanup);
-				appendPQExpBuffer(sql, "%sINDEX_CLEANUP FALSE", sep);
-				sep = comma;
-			}
-			if (vacopts->force_index_cleanup)
-			{
-				/* "INDEX_CLEANUP TRUE" has been supported since v12 */
-				Assert(serverVersion >= 120000);
-				Assert(!vacopts->no_index_cleanup);
-				appendPQExpBuffer(sql, "%sINDEX_CLEANUP TRUE", sep);
-				sep = comma;
-			}
-			if (!vacopts->do_truncate)
-			{
-				/* TRUNCATE is supported since v12 */
-				Assert(serverVersion >= 120000);
-				appendPQExpBuffer(sql, "%sTRUNCATE FALSE", sep);
-				sep = comma;
-			}
-			if (!vacopts->process_main)
-			{
-				/* PROCESS_MAIN is supported since v16 */
-				Assert(serverVersion >= 160000);
-				appendPQExpBuffer(sql, "%sPROCESS_MAIN FALSE", sep);
-				sep = comma;
-			}
-			if (!vacopts->process_toast)
-			{
-				/* PROCESS_TOAST is supported since v14 */
-				Assert(serverVersion >= 140000);
-				appendPQExpBuffer(sql, "%sPROCESS_TOAST FALSE", sep);
-				sep = comma;
-			}
-			if (vacopts->skip_database_stats)
-			{
-				/* SKIP_DATABASE_STATS is supported since v16 */
-				Assert(serverVersion >= 160000);
-				appendPQExpBuffer(sql, "%sSKIP_DATABASE_STATS", sep);
-				sep = comma;
-			}
-			if (vacopts->skip_locked)
-			{
-				/* SKIP_LOCKED is supported since v12 */
-				Assert(serverVersion >= 120000);
-				appendPQExpBuffer(sql, "%sSKIP_LOCKED", sep);
-				sep = comma;
-			}
-			if (vacopts->full)
-			{
-				appendPQExpBuffer(sql, "%sFULL", sep);
-				sep = comma;
-			}
-			if (vacopts->freeze)
-			{
-				appendPQExpBuffer(sql, "%sFREEZE", sep);
-				sep = comma;
-			}
-			if (vacopts->verbose)
-			{
-				appendPQExpBuffer(sql, "%sVERBOSE", sep);
-				sep = comma;
-			}
-			if (vacopts->and_analyze)
-			{
-				appendPQExpBuffer(sql, "%sANALYZE", sep);
-				sep = comma;
-			}
-			if (vacopts->parallel_workers >= 0)
-			{
-				/* PARALLEL is supported since v13 */
-				Assert(serverVersion >= 130000);
-				appendPQExpBuffer(sql, "%sPARALLEL %d", sep,
-								  vacopts->parallel_workers);
-				sep = comma;
-			}
-			if (vacopts->buffer_usage_limit)
-			{
-				Assert(serverVersion >= 160000);
-				appendPQExpBuffer(sql, "%sBUFFER_USAGE_LIMIT '%s'", sep,
-								  vacopts->buffer_usage_limit);
-				sep = comma;
-			}
-			if (sep != paren)
-				appendPQExpBufferChar(sql, ')');
-		}
-		else
-		{
-			if (vacopts->full)
-				appendPQExpBufferStr(sql, " FULL");
-			if (vacopts->freeze)
-				appendPQExpBufferStr(sql, " FREEZE");
-			if (vacopts->verbose)
-				appendPQExpBufferStr(sql, " VERBOSE");
-			if (vacopts->and_analyze)
-				appendPQExpBufferStr(sql, " ANALYZE");
-		}
-	}
-
-	appendPQExpBuffer(sql, " %s;", table);
-}
-
-/*
- * Send a vacuum/analyze command to the server, returning after sending the
- * command.
- *
- * Any errors during command execution are reported to stderr.
- */
-static void
-run_vacuum_command(PGconn *conn, const char *sql, bool echo,
-				   const char *table)
-{
-	bool		status;
-
-	if (echo)
-		printf("%s\n", sql);
-
-	status = PQsendQuery(conn, sql) == 1;
-
-	if (!status)
-	{
-		if (table)
-			pg_log_error("vacuuming of table \"%s\" in database \"%s\" failed: %s",
-						 table, PQdb(conn), PQerrorMessage(conn));
-		else
-			pg_log_error("vacuuming of database \"%s\" failed: %s",
-						 PQdb(conn), PQerrorMessage(conn));
-	}
-}
 
 static void
 help(const char *progname)
diff --git a/src/bin/scripts/vacuuming.c b/src/bin/scripts/vacuuming.c
new file mode 100644
index 00000000000..9be37fcc45a
--- /dev/null
+++ b/src/bin/scripts/vacuuming.c
@@ -0,0 +1,978 @@
+/*-------------------------------------------------------------------------
+ * vacuuming.c
+ *		Common routines for vacuumdb
+ *
+ * Portions Copyright (c) 1996-2025, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994, Regents of the University of California
+ *
+ * src/bin/scripts/vacuuming.c
+ *
+ *-------------------------------------------------------------------------
+ */
+
+#include "postgres_fe.h"
+
+#include <limits.h>
+
+#include "catalog/pg_attribute_d.h"
+#include "catalog/pg_class_d.h"
+#include "common/connect.h"
+#include "common/logging.h"
+#include "fe_utils/cancel.h"
+#include "fe_utils/option_utils.h"
+#include "fe_utils/parallel_slot.h"
+#include "fe_utils/query_utils.h"
+#include "fe_utils/string_utils.h"
+#include "vacuuming.h"
+
+VacObjFilter objfilter = OBJFILTER_NONE;
+
+
+/*
+ * Executes vacuum/analyze as indicated, or dies in case of failure.
+ */
+void
+vacuuming_main(ConnParams *cparams, const char *dbname,
+			   const char *maintenance_db, vacuumingOptions *vacopts,
+			   SimpleStringList *objects, bool analyze_in_stages,
+			   int tbl_count, int concurrentCons,
+			   const char *progname, bool echo, bool quiet)
+{
+	setup_cancel_handler(NULL);
+
+	/* Avoid opening extra connections. */
+	if (tbl_count && (concurrentCons > tbl_count))
+		concurrentCons = tbl_count;
+
+	if (objfilter & OBJFILTER_ALL_DBS)
+	{
+		cparams->dbname = maintenance_db;
+
+		vacuum_all_databases(cparams, vacopts,
+							 analyze_in_stages,
+							 objects,
+							 concurrentCons,
+							 progname, echo, quiet);
+	}
+	else
+	{
+		if (dbname == NULL)
+		{
+			if (getenv("PGDATABASE"))
+				dbname = getenv("PGDATABASE");
+			else if (getenv("PGUSER"))
+				dbname = getenv("PGUSER");
+			else
+				dbname = get_user_name_or_exit(progname);
+		}
+
+		cparams->dbname = dbname;
+
+		if (analyze_in_stages)
+		{
+			int			stage;
+			SimpleStringList *found_objs = NULL;
+
+			for (stage = 0; stage < ANALYZE_NUM_STAGES; stage++)
+			{
+				vacuum_one_database(cparams, vacopts,
+									stage,
+									objects,
+									vacopts->missing_stats_only ? &found_objs : NULL,
+									concurrentCons,
+									progname, echo, quiet);
+			}
+		}
+		else
+			vacuum_one_database(cparams, vacopts,
+								ANALYZE_NO_STAGE,
+								objects, NULL,
+								concurrentCons,
+								progname, echo, quiet);
+	}
+}
+
+
+/*
+ * vacuum_one_database
+ *
+ * Process tables in the given database.
+ *
+ * There are two ways to specify the list of objects to process:
+ *
+ * 1) The "found_objs" parameter is a double pointer to a fully qualified list
+ *    of objects to process, as returned by a previous call to
+ *    vacuum_one_database().
+ *
+ *     a) If both "found_objs" (the double pointer) and "*found_objs" (the
+ *        once-dereferenced double pointer) are not NULL, this list takes
+ *        priority, and anything specified in "objects" is ignored.
+ *
+ *     b) If "found_objs" (the double pointer) is not NULL but "*found_objs"
+ *        (the once-dereferenced double pointer) _is_ NULL, the "objects"
+ *        parameter takes priority, and the results of the catalog query
+ *        described in (2) are stored in "found_objs".
+ *
+ *     c) If "found_objs" (the double pointer) is NULL, the "objects"
+ *        parameter again takes priority, and the results of the catalog query
+ *        are not saved.
+ *
+ * 2) The "objects" parameter is a user-specified list of objects to process.
+ *    When (1b) or (1c) applies, this function performs a catalog query to
+ *    retrieve a fully qualified list of objects to process, as described
+ *    below.
+ *
+ *     a) If "objects" is not NULL, the catalog query gathers only the objects
+ *        listed in "objects".
+ *
+ *     b) If "objects" is NULL, all tables in the database are gathered.
+ *
+ * Note that this function is only concerned with running exactly one stage
+ * when in analyze-in-stages mode; caller must iterate on us if necessary.
+ *
+ * If concurrentCons is > 1, multiple connections are used to vacuum tables
+ * in parallel.
+ */
+void
+vacuum_one_database(ConnParams *cparams,
+					vacuumingOptions *vacopts,
+					int stage,
+					SimpleStringList *objects,
+					SimpleStringList **found_objs,
+					int concurrentCons,
+					const char *progname, bool echo, bool quiet)
+{
+	PQExpBufferData sql;
+	PGconn	   *conn;
+	SimpleStringListCell *cell;
+	ParallelSlotArray *sa;
+	int			ntups = 0;
+	bool		failed = false;
+	const char *initcmd;
+	SimpleStringList *ret = NULL;
+	const char *stage_commands[] = {
+		"SET default_statistics_target=1; SET vacuum_cost_delay=0;",
+		"SET default_statistics_target=10; RESET vacuum_cost_delay;",
+		"RESET default_statistics_target;"
+	};
+	const char *stage_messages[] = {
+		gettext_noop("Generating minimal optimizer statistics (1 target)"),
+		gettext_noop("Generating medium optimizer statistics (10 targets)"),
+		gettext_noop("Generating default (full) optimizer statistics")
+	};
+
+	Assert(stage == ANALYZE_NO_STAGE ||
+		   (stage >= 0 && stage < ANALYZE_NUM_STAGES));
+
+	conn = connectDatabase(cparams, progname, echo, false, true);
+
+	if (vacopts->disable_page_skipping && PQserverVersion(conn) < 90600)
+	{
+		PQfinish(conn);
+		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
+				 "disable-page-skipping", "9.6");
+	}
+
+	if (vacopts->no_index_cleanup && PQserverVersion(conn) < 120000)
+	{
+		PQfinish(conn);
+		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
+				 "no-index-cleanup", "12");
+	}
+
+	if (vacopts->force_index_cleanup && PQserverVersion(conn) < 120000)
+	{
+		PQfinish(conn);
+		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
+				 "force-index-cleanup", "12");
+	}
+
+	if (!vacopts->do_truncate && PQserverVersion(conn) < 120000)
+	{
+		PQfinish(conn);
+		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
+				 "no-truncate", "12");
+	}
+
+	if (!vacopts->process_main && PQserverVersion(conn) < 160000)
+	{
+		PQfinish(conn);
+		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
+				 "no-process-main", "16");
+	}
+
+	if (!vacopts->process_toast && PQserverVersion(conn) < 140000)
+	{
+		PQfinish(conn);
+		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
+				 "no-process-toast", "14");
+	}
+
+	if (vacopts->skip_locked && PQserverVersion(conn) < 120000)
+	{
+		PQfinish(conn);
+		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
+				 "skip-locked", "12");
+	}
+
+	if (vacopts->min_xid_age != 0 && PQserverVersion(conn) < 90600)
+	{
+		PQfinish(conn);
+		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
+				 "--min-xid-age", "9.6");
+	}
+
+	if (vacopts->min_mxid_age != 0 && PQserverVersion(conn) < 90600)
+	{
+		PQfinish(conn);
+		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
+				 "--min-mxid-age", "9.6");
+	}
+
+	if (vacopts->parallel_workers >= 0 && PQserverVersion(conn) < 130000)
+	{
+		PQfinish(conn);
+		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
+				 "--parallel", "13");
+	}
+
+	if (vacopts->buffer_usage_limit && PQserverVersion(conn) < 160000)
+	{
+		PQfinish(conn);
+		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
+				 "--buffer-usage-limit", "16");
+	}
+
+	if (vacopts->missing_stats_only && PQserverVersion(conn) < 150000)
+	{
+		PQfinish(conn);
+		pg_fatal("cannot use the \"%s\" option on server versions older than PostgreSQL %s",
+				 "--missing-stats-only", "15");
+	}
+
+	/* skip_database_stats is used automatically if server supports it */
+	vacopts->skip_database_stats = (PQserverVersion(conn) >= 160000);
+
+	if (!quiet)
+	{
+		if (stage != ANALYZE_NO_STAGE)
+			printf(_("%s: processing database \"%s\": %s\n"),
+				   progname, PQdb(conn), _(stage_messages[stage]));
+		else
+			printf(_("%s: vacuuming database \"%s\"\n"),
+				   progname, PQdb(conn));
+		fflush(stdout);
+	}
+
+	/*
+	 * If the caller provided the results of a previous catalog query, just
+	 * use that.  Otherwise, run the catalog query ourselves and set the
+	 * return variable if provided.
+	 */
+	if (found_objs && *found_objs)
+		ret = *found_objs;
+	else
+	{
+		ret = retrieve_objects(conn, vacopts, objects, echo);
+		if (found_objs)
+			*found_objs = ret;
+	}
+
+	/*
+	 * Count the number of objects in the catalog query result.  If there are
+	 * none, we are done.
+	 */
+	for (cell = ret ? ret->head : NULL; cell; cell = cell->next)
+		ntups++;
+
+	if (ntups == 0)
+	{
+		PQfinish(conn);
+		return;
+	}
+
+	/*
+	 * Ensure concurrentCons is sane.  If there are more connections than
+	 * vacuumable relations, we don't need to use them all.
+	 */
+	if (concurrentCons > ntups)
+		concurrentCons = ntups;
+	if (concurrentCons <= 0)
+		concurrentCons = 1;
+
+	/*
+	 * All slots need to be prepared to run the appropriate analyze stage, if
+	 * caller requested that mode.  We have to prepare the initial connection
+	 * ourselves before setting up the slots.
+	 */
+	if (stage == ANALYZE_NO_STAGE)
+		initcmd = NULL;
+	else
+	{
+		initcmd = stage_commands[stage];
+		executeCommand(conn, initcmd, echo);
+	}
+
+	/*
+	 * Setup the database connections. We reuse the connection we already have
+	 * for the first slot.  If not in parallel mode, the first slot in the
+	 * array contains the connection.
+	 */
+	sa = ParallelSlotsSetup(concurrentCons, cparams, progname, echo, initcmd);
+	ParallelSlotsAdoptConn(sa, conn);
+
+	initPQExpBuffer(&sql);
+
+	cell = ret->head;
+	do
+	{
+		const char *tabname = cell->val;
+		ParallelSlot *free_slot;
+
+		if (CancelRequested)
+		{
+			failed = true;
+			goto finish;
+		}
+
+		free_slot = ParallelSlotsGetIdle(sa, NULL);
+		if (!free_slot)
+		{
+			failed = true;
+			goto finish;
+		}
+
+		prepare_vacuum_command(&sql, PQserverVersion(free_slot->connection),
+							   vacopts, tabname);
+
+		/*
+		 * Execute the vacuum.  All errors are handled in processQueryResult
+		 * through ParallelSlotsGetIdle.
+		 */
+		ParallelSlotSetHandler(free_slot, TableCommandResultHandler, NULL);
+		run_vacuum_command(free_slot->connection, sql.data,
+						   echo, tabname);
+
+		cell = cell->next;
+	} while (cell != NULL);
+
+	if (!ParallelSlotsWaitCompletion(sa))
+	{
+		failed = true;
+		goto finish;
+	}
+
+	/* If we used SKIP_DATABASE_STATS, mop up with ONLY_DATABASE_STATS */
+	if (vacopts->skip_database_stats &&
+		stage == ANALYZE_NO_STAGE)
+	{
+		const char *cmd = "VACUUM (ONLY_DATABASE_STATS);";
+		ParallelSlot *free_slot = ParallelSlotsGetIdle(sa, NULL);
+
+		if (!free_slot)
+		{
+			failed = true;
+			goto finish;
+		}
+
+		ParallelSlotSetHandler(free_slot, TableCommandResultHandler, NULL);
+		run_vacuum_command(free_slot->connection, cmd, echo, NULL);
+
+		if (!ParallelSlotsWaitCompletion(sa))
+			failed = true;
+	}
+
+finish:
+	ParallelSlotsTerminate(sa);
+	pg_free(sa);
+
+	termPQExpBuffer(&sql);
+
+	if (failed)
+		exit(1);
+}
+
+/*
+ * Prepare the list of tables to process by querying the catalogs.
+ *
+ * Since we execute the constructed query with the default search_path (which
+ * could be unsafe), everything in this query MUST be fully qualified.
+ *
+ * First, build a WITH clause for the catalog query if any tables were
+ * specified, with a set of values made of relation names and their optional
+ * set of columns.  This is used to match any provided column lists with the
+ * generated qualified identifiers and to filter for the tables provided via
+ * --table.  If a listed table does not exist, the catalog query will fail.
+ */
+SimpleStringList *
+retrieve_objects(PGconn *conn, vacuumingOptions *vacopts,
+				 SimpleStringList *objects, bool echo)
+{
+	PQExpBufferData buf;
+	PQExpBufferData catalog_query;
+	PGresult   *res;
+	SimpleStringListCell *cell;
+	SimpleStringList *found_objs = palloc0(sizeof(SimpleStringList));
+	bool		objects_listed = false;
+
+	initPQExpBuffer(&catalog_query);
+	for (cell = objects ? objects->head : NULL; cell; cell = cell->next)
+	{
+		char	   *just_table = NULL;
+		const char *just_columns = NULL;
+
+		if (!objects_listed)
+		{
+			appendPQExpBufferStr(&catalog_query,
+								 "WITH listed_objects (object_oid, column_list) AS (\n"
+								 "  VALUES (");
+			objects_listed = true;
+		}
+		else
+			appendPQExpBufferStr(&catalog_query, ",\n  (");
+
+		if (objfilter & (OBJFILTER_SCHEMA | OBJFILTER_SCHEMA_EXCLUDE))
+		{
+			appendStringLiteralConn(&catalog_query, cell->val, conn);
+			appendPQExpBufferStr(&catalog_query, "::pg_catalog.regnamespace, ");
+		}
+
+		if (objfilter & OBJFILTER_TABLE)
+		{
+			/*
+			 * Split relation and column names given by the user, this is used
+			 * to feed the CTE with values on which are performed pre-run
+			 * validity checks as well.  For now these happen only on the
+			 * relation name.
+			 */
+			splitTableColumnsSpec(cell->val, PQclientEncoding(conn),
+								  &just_table, &just_columns);
+
+			appendStringLiteralConn(&catalog_query, just_table, conn);
+			appendPQExpBufferStr(&catalog_query, "::pg_catalog.regclass, ");
+		}
+
+		if (just_columns && just_columns[0] != '\0')
+			appendStringLiteralConn(&catalog_query, just_columns, conn);
+		else
+			appendPQExpBufferStr(&catalog_query, "NULL");
+
+		appendPQExpBufferStr(&catalog_query, "::pg_catalog.text)");
+
+		pg_free(just_table);
+	}
+
+	/* Finish formatting the CTE */
+	if (objects_listed)
+		appendPQExpBufferStr(&catalog_query, "\n)\n");
+
+	appendPQExpBufferStr(&catalog_query, "SELECT c.relname, ns.nspname");
+
+	if (objects_listed)
+		appendPQExpBufferStr(&catalog_query, ", listed_objects.column_list");
+
+	appendPQExpBufferStr(&catalog_query,
+						 " FROM pg_catalog.pg_class c\n"
+						 " JOIN pg_catalog.pg_namespace ns"
+						 " ON c.relnamespace OPERATOR(pg_catalog.=) ns.oid\n"
+						 " CROSS JOIN LATERAL (SELECT c.relkind IN ("
+						 CppAsString2(RELKIND_PARTITIONED_TABLE) ", "
+						 CppAsString2(RELKIND_PARTITIONED_INDEX) ")) as p (inherited)\n"
+						 " LEFT JOIN pg_catalog.pg_class t"
+						 " ON c.reltoastrelid OPERATOR(pg_catalog.=) t.oid\n");
+
+	/*
+	 * Used to match the tables or schemas listed by the user, completing the
+	 * JOIN clause.
+	 */
+	if (objects_listed)
+	{
+		appendPQExpBufferStr(&catalog_query, " LEFT JOIN listed_objects"
+							 " ON listed_objects.object_oid"
+							 " OPERATOR(pg_catalog.=) ");
+
+		if (objfilter & OBJFILTER_TABLE)
+			appendPQExpBufferStr(&catalog_query, "c.oid\n");
+		else
+			appendPQExpBufferStr(&catalog_query, "ns.oid\n");
+	}
+
+	/*
+	 * Exclude temporary tables, beginning the WHERE clause.
+	 */
+	appendPQExpBufferStr(&catalog_query,
+						 " WHERE c.relpersistence OPERATOR(pg_catalog.!=) "
+						 CppAsString2(RELPERSISTENCE_TEMP) "\n");
+
+	/*
+	 * Used to match the tables or schemas listed by the user, for the WHERE
+	 * clause.
+	 */
+	if (objects_listed)
+	{
+		if (objfilter & OBJFILTER_SCHEMA_EXCLUDE)
+			appendPQExpBufferStr(&catalog_query,
+								 " AND listed_objects.object_oid IS NULL\n");
+		else
+			appendPQExpBufferStr(&catalog_query,
+								 " AND listed_objects.object_oid IS NOT NULL\n");
+	}
+
+	/*
+	 * If no tables were listed, filter for the relevant relation types.  If
+	 * tables were given via --table, don't bother filtering by relation type.
+	 * Instead, let the server decide whether a given relation can be
+	 * processed in which case the user will know about it.
+	 */
+	if ((objfilter & OBJFILTER_TABLE) == 0)
+	{
+		/*
+		 * vacuumdb should generally follow the behavior of the underlying
+		 * VACUUM and ANALYZE commands. If analyze_only is true, process
+		 * regular tables, materialized views, and partitioned tables, just
+		 * like ANALYZE (with no specific target tables) does. Otherwise,
+		 * process only regular tables and materialized views, since VACUUM
+		 * skips partitioned tables when no target tables are specified.
+		 */
+		if (vacopts->analyze_only)
+			appendPQExpBufferStr(&catalog_query,
+								 " AND c.relkind OPERATOR(pg_catalog.=) ANY (array["
+								 CppAsString2(RELKIND_RELATION) ", "
+								 CppAsString2(RELKIND_MATVIEW) ", "
+								 CppAsString2(RELKIND_PARTITIONED_TABLE) "])\n");
+		else
+			appendPQExpBufferStr(&catalog_query,
+								 " AND c.relkind OPERATOR(pg_catalog.=) ANY (array["
+								 CppAsString2(RELKIND_RELATION) ", "
+								 CppAsString2(RELKIND_MATVIEW) "])\n");
+	}
+
+	/*
+	 * For --min-xid-age and --min-mxid-age, the age of the relation is the
+	 * greatest of the ages of the main relation and its associated TOAST
+	 * table.  The commands generated by vacuumdb will also process the TOAST
+	 * table for the relation if necessary, so it does not need to be
+	 * considered separately.
+	 */
+	if (vacopts->min_xid_age != 0)
+	{
+		appendPQExpBuffer(&catalog_query,
+						  " AND GREATEST(pg_catalog.age(c.relfrozenxid),"
+						  " pg_catalog.age(t.relfrozenxid)) "
+						  " OPERATOR(pg_catalog.>=) '%d'::pg_catalog.int4\n"
+						  " AND c.relfrozenxid OPERATOR(pg_catalog.!=)"
+						  " '0'::pg_catalog.xid\n",
+						  vacopts->min_xid_age);
+	}
+
+	if (vacopts->min_mxid_age != 0)
+	{
+		appendPQExpBuffer(&catalog_query,
+						  " AND GREATEST(pg_catalog.mxid_age(c.relminmxid),"
+						  " pg_catalog.mxid_age(t.relminmxid)) OPERATOR(pg_catalog.>=)"
+						  " '%d'::pg_catalog.int4\n"
+						  " AND c.relminmxid OPERATOR(pg_catalog.!=)"
+						  " '0'::pg_catalog.xid\n",
+						  vacopts->min_mxid_age);
+	}
+
+	if (vacopts->missing_stats_only)
+	{
+		appendPQExpBufferStr(&catalog_query, " AND (\n");
+
+		/* regular stats */
+		appendPQExpBufferStr(&catalog_query,
+							 " EXISTS (SELECT NULL FROM pg_catalog.pg_attribute a\n"
+							 " WHERE a.attrelid OPERATOR(pg_catalog.=) c.oid\n"
+							 " AND a.attnum OPERATOR(pg_catalog.>) 0::pg_catalog.int2\n"
+							 " AND NOT a.attisdropped\n"
+							 " AND a.attstattarget IS DISTINCT FROM 0::pg_catalog.int2\n"
+							 " AND a.attgenerated OPERATOR(pg_catalog.<>) "
+							 CppAsString2(ATTRIBUTE_GENERATED_VIRTUAL) "\n"
+							 " AND NOT EXISTS (SELECT NULL FROM pg_catalog.pg_statistic s\n"
+							 " WHERE s.starelid OPERATOR(pg_catalog.=) a.attrelid\n"
+							 " AND s.staattnum OPERATOR(pg_catalog.=) a.attnum\n"
+							 " AND s.stainherit OPERATOR(pg_catalog.=) p.inherited))\n");
+
+		/* extended stats */
+		appendPQExpBufferStr(&catalog_query,
+							 " OR EXISTS (SELECT NULL FROM pg_catalog.pg_statistic_ext e\n"
+							 " WHERE e.stxrelid OPERATOR(pg_catalog.=) c.oid\n"
+							 " AND e.stxstattarget IS DISTINCT FROM 0::pg_catalog.int2\n"
+							 " AND NOT EXISTS (SELECT NULL FROM pg_catalog.pg_statistic_ext_data d\n"
+							 " WHERE d.stxoid OPERATOR(pg_catalog.=) e.oid\n"
+							 " AND d.stxdinherit OPERATOR(pg_catalog.=) p.inherited))\n");
+
+		/* expression indexes */
+		appendPQExpBufferStr(&catalog_query,
+							 " OR EXISTS (SELECT NULL FROM pg_catalog.pg_attribute a\n"
+							 " JOIN pg_catalog.pg_index i"
+							 " ON i.indexrelid OPERATOR(pg_catalog.=) a.attrelid\n"
+							 " WHERE i.indrelid OPERATOR(pg_catalog.=) c.oid\n"
+							 " AND i.indkey[a.attnum OPERATOR(pg_catalog.-) 1::pg_catalog.int2]"
+							 " OPERATOR(pg_catalog.=) 0::pg_catalog.int2\n"
+							 " AND a.attnum OPERATOR(pg_catalog.>) 0::pg_catalog.int2\n"
+							 " AND NOT a.attisdropped\n"
+							 " AND a.attstattarget IS DISTINCT FROM 0::pg_catalog.int2\n"
+							 " AND NOT EXISTS (SELECT NULL FROM pg_catalog.pg_statistic s\n"
+							 " WHERE s.starelid OPERATOR(pg_catalog.=) a.attrelid\n"
+							 " AND s.staattnum OPERATOR(pg_catalog.=) a.attnum\n"
+							 " AND s.stainherit OPERATOR(pg_catalog.=) p.inherited))\n");
+
+		/* inheritance and regular stats */
+		appendPQExpBufferStr(&catalog_query,
+							 " OR EXISTS (SELECT NULL FROM pg_catalog.pg_attribute a\n"
+							 " WHERE a.attrelid OPERATOR(pg_catalog.=) c.oid\n"
+							 " AND a.attnum OPERATOR(pg_catalog.>) 0::pg_catalog.int2\n"
+							 " AND NOT a.attisdropped\n"
+							 " AND a.attstattarget IS DISTINCT FROM 0::pg_catalog.int2\n"
+							 " AND a.attgenerated OPERATOR(pg_catalog.<>) "
+							 CppAsString2(ATTRIBUTE_GENERATED_VIRTUAL) "\n"
+							 " AND c.relhassubclass\n"
+							 " AND NOT p.inherited\n"
+							 " AND EXISTS (SELECT NULL FROM pg_catalog.pg_inherits h\n"
+							 " WHERE h.inhparent OPERATOR(pg_catalog.=) c.oid)\n"
+							 " AND NOT EXISTS (SELECT NULL FROM pg_catalog.pg_statistic s\n"
+							 " WHERE s.starelid OPERATOR(pg_catalog.=) a.attrelid\n"
+							 " AND s.staattnum OPERATOR(pg_catalog.=) a.attnum\n"
+							 " AND s.stainherit))\n");
+
+		/* inheritance and extended stats */
+		appendPQExpBufferStr(&catalog_query,
+							 " OR EXISTS (SELECT NULL FROM pg_catalog.pg_statistic_ext e\n"
+							 " WHERE e.stxrelid OPERATOR(pg_catalog.=) c.oid\n"
+							 " AND e.stxstattarget IS DISTINCT FROM 0::pg_catalog.int2\n"
+							 " AND c.relhassubclass\n"
+							 " AND NOT p.inherited\n"
+							 " AND EXISTS (SELECT NULL FROM pg_catalog.pg_inherits h\n"
+							 " WHERE h.inhparent OPERATOR(pg_catalog.=) c.oid)\n"
+							 " AND NOT EXISTS (SELECT NULL FROM pg_catalog.pg_statistic_ext_data d\n"
+							 " WHERE d.stxoid OPERATOR(pg_catalog.=) e.oid\n"
+							 " AND d.stxdinherit))\n");
+
+		appendPQExpBufferStr(&catalog_query, " )\n");
+	}
+
+	/*
+	 * Execute the catalog query.  We use the default search_path for this
+	 * query for consistency with table lookups done elsewhere by the user.
+	 */
+	appendPQExpBufferStr(&catalog_query, " ORDER BY c.relpages DESC;");
+	executeCommand(conn, "RESET search_path;", echo);
+	res = executeQuery(conn, catalog_query.data, echo);
+	termPQExpBuffer(&catalog_query);
+	PQclear(executeQuery(conn, ALWAYS_SECURE_SEARCH_PATH_SQL, echo));
+
+	/*
+	 * Build qualified identifiers for each table, including the column list
+	 * if given.
+	 */
+	initPQExpBuffer(&buf);
+	for (int i = 0; i < PQntuples(res); i++)
+	{
+		appendPQExpBufferStr(&buf,
+							 fmtQualifiedIdEnc(PQgetvalue(res, i, 1),
+											   PQgetvalue(res, i, 0),
+											   PQclientEncoding(conn)));
+
+		if (objects_listed && !PQgetisnull(res, i, 2))
+			appendPQExpBufferStr(&buf, PQgetvalue(res, i, 2));
+
+		simple_string_list_append(found_objs, buf.data);
+		resetPQExpBuffer(&buf);
+	}
+	termPQExpBuffer(&buf);
+	PQclear(res);
+
+	return found_objs;
+}
+
+/*
+ * Vacuum/analyze all connectable databases.
+ *
+ * In analyze-in-stages mode, we process all databases in one stage before
+ * moving on to the next stage.  That ensure minimal stats are available
+ * quickly everywhere before generating more detailed ones.
+ */
+void
+vacuum_all_databases(ConnParams *cparams,
+					 vacuumingOptions *vacopts,
+					 bool analyze_in_stages,
+					 SimpleStringList *objects,
+					 int concurrentCons,
+					 const char *progname, bool echo, bool quiet)
+{
+	PGconn	   *conn;
+	PGresult   *result;
+	int			stage;
+	int			i;
+
+	conn = connectMaintenanceDatabase(cparams, progname, echo);
+	result = executeQuery(conn,
+						  "SELECT datname FROM pg_database WHERE datallowconn AND datconnlimit <> -2 ORDER BY 1;",
+						  echo);
+	PQfinish(conn);
+
+	if (analyze_in_stages)
+	{
+		SimpleStringList **found_objs = NULL;
+
+		if (vacopts->missing_stats_only)
+			found_objs = palloc0(PQntuples(result) * sizeof(SimpleStringList *));
+
+		/*
+		 * When analyzing all databases in stages, we analyze them all in the
+		 * fastest stage first, so that initial statistics become available
+		 * for all of them as soon as possible.
+		 *
+		 * This means we establish several times as many connections, but
+		 * that's a secondary consideration.
+		 */
+		for (stage = 0; stage < ANALYZE_NUM_STAGES; stage++)
+		{
+			for (i = 0; i < PQntuples(result); i++)
+			{
+				cparams->override_dbname = PQgetvalue(result, i, 0);
+
+				vacuum_one_database(cparams, vacopts,
+									stage,
+									objects,
+									vacopts->missing_stats_only ? &found_objs[i] : NULL,
+									concurrentCons,
+									progname, echo, quiet);
+			}
+		}
+	}
+	else
+	{
+		for (i = 0; i < PQntuples(result); i++)
+		{
+			cparams->override_dbname = PQgetvalue(result, i, 0);
+
+			vacuum_one_database(cparams, vacopts,
+								ANALYZE_NO_STAGE,
+								objects, NULL,
+								concurrentCons,
+								progname, echo, quiet);
+		}
+	}
+
+	PQclear(result);
+}
+
+/*
+ * Construct a vacuum/analyze command to run based on the given
+ * options, in the given string buffer, which may contain previous garbage.
+ *
+ * The table name used must be already properly quoted.  The command generated
+ * depends on the server version involved and it is semicolon-terminated.
+ */
+void
+prepare_vacuum_command(PQExpBuffer sql, int serverVersion,
+					   vacuumingOptions *vacopts, const char *table)
+{
+	const char *paren = " (";
+	const char *comma = ", ";
+	const char *sep = paren;
+
+	resetPQExpBuffer(sql);
+
+	if (vacopts->analyze_only)
+	{
+		appendPQExpBufferStr(sql, "ANALYZE");
+
+		/* parenthesized grammar of ANALYZE is supported since v11 */
+		if (serverVersion >= 110000)
+		{
+			if (vacopts->skip_locked)
+			{
+				/* SKIP_LOCKED is supported since v12 */
+				Assert(serverVersion >= 120000);
+				appendPQExpBuffer(sql, "%sSKIP_LOCKED", sep);
+				sep = comma;
+			}
+			if (vacopts->verbose)
+			{
+				appendPQExpBuffer(sql, "%sVERBOSE", sep);
+				sep = comma;
+			}
+			if (vacopts->buffer_usage_limit)
+			{
+				Assert(serverVersion >= 160000);
+				appendPQExpBuffer(sql, "%sBUFFER_USAGE_LIMIT '%s'", sep,
+								  vacopts->buffer_usage_limit);
+				sep = comma;
+			}
+			if (sep != paren)
+				appendPQExpBufferChar(sql, ')');
+		}
+		else
+		{
+			if (vacopts->verbose)
+				appendPQExpBufferStr(sql, " VERBOSE");
+		}
+	}
+	else
+	{
+		appendPQExpBufferStr(sql, "VACUUM");
+
+		/* parenthesized grammar of VACUUM is supported since v9.0 */
+		if (serverVersion >= 90000)
+		{
+			if (vacopts->disable_page_skipping)
+			{
+				/* DISABLE_PAGE_SKIPPING is supported since v9.6 */
+				Assert(serverVersion >= 90600);
+				appendPQExpBuffer(sql, "%sDISABLE_PAGE_SKIPPING", sep);
+				sep = comma;
+			}
+			if (vacopts->no_index_cleanup)
+			{
+				/* "INDEX_CLEANUP FALSE" has been supported since v12 */
+				Assert(serverVersion >= 120000);
+				Assert(!vacopts->force_index_cleanup);
+				appendPQExpBuffer(sql, "%sINDEX_CLEANUP FALSE", sep);
+				sep = comma;
+			}
+			if (vacopts->force_index_cleanup)
+			{
+				/* "INDEX_CLEANUP TRUE" has been supported since v12 */
+				Assert(serverVersion >= 120000);
+				Assert(!vacopts->no_index_cleanup);
+				appendPQExpBuffer(sql, "%sINDEX_CLEANUP TRUE", sep);
+				sep = comma;
+			}
+			if (!vacopts->do_truncate)
+			{
+				/* TRUNCATE is supported since v12 */
+				Assert(serverVersion >= 120000);
+				appendPQExpBuffer(sql, "%sTRUNCATE FALSE", sep);
+				sep = comma;
+			}
+			if (!vacopts->process_main)
+			{
+				/* PROCESS_MAIN is supported since v16 */
+				Assert(serverVersion >= 160000);
+				appendPQExpBuffer(sql, "%sPROCESS_MAIN FALSE", sep);
+				sep = comma;
+			}
+			if (!vacopts->process_toast)
+			{
+				/* PROCESS_TOAST is supported since v14 */
+				Assert(serverVersion >= 140000);
+				appendPQExpBuffer(sql, "%sPROCESS_TOAST FALSE", sep);
+				sep = comma;
+			}
+			if (vacopts->skip_database_stats)
+			{
+				/* SKIP_DATABASE_STATS is supported since v16 */
+				Assert(serverVersion >= 160000);
+				appendPQExpBuffer(sql, "%sSKIP_DATABASE_STATS", sep);
+				sep = comma;
+			}
+			if (vacopts->skip_locked)
+			{
+				/* SKIP_LOCKED is supported since v12 */
+				Assert(serverVersion >= 120000);
+				appendPQExpBuffer(sql, "%sSKIP_LOCKED", sep);
+				sep = comma;
+			}
+			if (vacopts->full)
+			{
+				appendPQExpBuffer(sql, "%sFULL", sep);
+				sep = comma;
+			}
+			if (vacopts->freeze)
+			{
+				appendPQExpBuffer(sql, "%sFREEZE", sep);
+				sep = comma;
+			}
+			if (vacopts->verbose)
+			{
+				appendPQExpBuffer(sql, "%sVERBOSE", sep);
+				sep = comma;
+			}
+			if (vacopts->and_analyze)
+			{
+				appendPQExpBuffer(sql, "%sANALYZE", sep);
+				sep = comma;
+			}
+			if (vacopts->parallel_workers >= 0)
+			{
+				/* PARALLEL is supported since v13 */
+				Assert(serverVersion >= 130000);
+				appendPQExpBuffer(sql, "%sPARALLEL %d", sep,
+								  vacopts->parallel_workers);
+				sep = comma;
+			}
+			if (vacopts->buffer_usage_limit)
+			{
+				Assert(serverVersion >= 160000);
+				appendPQExpBuffer(sql, "%sBUFFER_USAGE_LIMIT '%s'", sep,
+								  vacopts->buffer_usage_limit);
+				sep = comma;
+			}
+			if (sep != paren)
+				appendPQExpBufferChar(sql, ')');
+		}
+		else
+		{
+			if (vacopts->full)
+				appendPQExpBufferStr(sql, " FULL");
+			if (vacopts->freeze)
+				appendPQExpBufferStr(sql, " FREEZE");
+			if (vacopts->verbose)
+				appendPQExpBufferStr(sql, " VERBOSE");
+			if (vacopts->and_analyze)
+				appendPQExpBufferStr(sql, " ANALYZE");
+		}
+	}
+
+	appendPQExpBuffer(sql, " %s;", table);
+}
+
+/*
+ * Send a vacuum/analyze command to the server, returning after sending the
+ * command.
+ *
+ * Any errors during command execution are reported to stderr.
+ */
+void
+run_vacuum_command(PGconn *conn, const char *sql, bool echo,
+				   const char *table)
+{
+	bool		status;
+
+	if (echo)
+		printf("%s\n", sql);
+
+	status = PQsendQuery(conn, sql) == 1;
+
+	if (!status)
+	{
+		if (table)
+		{
+			pg_log_error("vacuuming of table \"%s\" in database \"%s\" failed: %s",
+						 table, PQdb(conn), PQerrorMessage(conn));
+		}
+		else
+		{
+			pg_log_error("vacuuming of database \"%s\" failed: %s",
+						 PQdb(conn), PQerrorMessage(conn));
+		}
+	}
+}
+
+/*
+ * Returns a newly malloc'd version of 'src' with escaped single quotes and
+ * backslashes.
+ */
+char *
+escape_quotes(const char *src)
+{
+	char	   *result = escape_single_quotes_ascii(src);
+
+	if (!result)
+		pg_fatal("out of memory");
+	return result;
+}
diff --git a/src/bin/scripts/vacuuming.h b/src/bin/scripts/vacuuming.h
new file mode 100644
index 00000000000..d3f000840fa
--- /dev/null
+++ b/src/bin/scripts/vacuuming.h
@@ -0,0 +1,95 @@
+/*-------------------------------------------------------------------------
+ *
+ * vacuuming.h
+ *		Common declarations for vacuuming.c
+ *
+ * Portions Copyright (c) 1996-2025, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994, Regents of the University of California
+ *
+ * src/bin/scripts/vacuuming.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef VACUUMING_H
+#define VACUUMING_H
+
+#include "common.h"
+#include "fe_utils/connect_utils.h"
+#include "fe_utils/simple_list.h"
+
+/* For analyze-in-stages mode */
+#define ANALYZE_NO_STAGE	-1
+#define ANALYZE_NUM_STAGES	3
+
+/* vacuum options controlled by user flags */
+typedef struct vacuumingOptions
+{
+	bool		analyze_only;
+	bool		verbose;
+	bool		and_analyze;
+	bool		full;
+	bool		freeze;
+	bool		disable_page_skipping;
+	bool		skip_locked;
+	int			min_xid_age;
+	int			min_mxid_age;
+	int			parallel_workers;	/* >= 0 indicates user specified the
+									 * parallel degree, otherwise -1 */
+	bool		no_index_cleanup;
+	bool		force_index_cleanup;
+	bool		do_truncate;
+	bool		process_main;
+	bool		process_toast;
+	bool		skip_database_stats;
+	char	   *buffer_usage_limit;
+	bool		missing_stats_only;
+} vacuumingOptions;
+
+/* object filter options */
+typedef enum
+{
+	OBJFILTER_NONE = 0,			/* no filter used */
+	OBJFILTER_ALL_DBS = (1 << 0),	/* -a | --all */
+	OBJFILTER_DATABASE = (1 << 1),	/* -d | --dbname */
+	OBJFILTER_TABLE = (1 << 2), /* -t | --table */
+	OBJFILTER_SCHEMA = (1 << 3),	/* -n | --schema */
+	OBJFILTER_SCHEMA_EXCLUDE = (1 << 4),	/* -N | --exclude-schema */
+} VacObjFilter;
+
+extern VacObjFilter objfilter;
+
+extern void vacuuming_main(ConnParams *cparams, const char *dbname,
+						   const char *maintenance_db, vacuumingOptions *vacopts,
+						   SimpleStringList *objects, bool analyze_in_stages,
+						   int tbl_count, int concurrentCons,
+						   const char *progname, bool echo, bool quiet);
+
+extern SimpleStringList *retrieve_objects(PGconn *conn,
+										  vacuumingOptions *vacopts,
+										  SimpleStringList *objects,
+										  bool echo);
+
+extern void vacuum_one_database(ConnParams *cparams,
+								vacuumingOptions *vacopts,
+								int stage,
+								SimpleStringList *objects,
+								SimpleStringList **found_objs,
+								int concurrentCons,
+								const char *progname, bool echo, bool quiet);
+
+extern void vacuum_all_databases(ConnParams *cparams,
+								 vacuumingOptions *vacopts,
+								 bool analyze_in_stages,
+								 SimpleStringList *objects,
+								 int concurrentCons,
+								 const char *progname, bool echo, bool quiet);
+
+extern void prepare_vacuum_command(PQExpBuffer sql, int serverVersion,
+								   vacuumingOptions *vacopts, const char *table);
+
+extern void run_vacuum_command(PGconn *conn, const char *sql, bool echo,
+							   const char *table);
+
+extern char *escape_quotes(const char *src);
+
+#endif							/* VACUUMING_H */
-- 
2.39.5


--ipmcfwv2cm3fpxaf
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v19-0002-Add-REPACK-command.patch"



^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
  0 siblings, 0 replies; 315+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 315+ messages in thread


end of thread, other threads:[~2026-04-17 13:13 UTC | newest]

Thread overview: 315+ messages (download: mbox mbox.gz follow: Atom feed)
-- links below jump to the message on this page --
2025-08-30 12:39 [PATCH v19 1/6] Split vacuumdb to create vacuuming.c/h Álvaro Herrera <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>

This inbox is served by agora; see mirroring instructions
for how to clone and mirror all data and code used for this inbox